agora inbox for [email protected]
help / color / mirror / Atom feed[PATCH v45 5/8] rename cluster.c/h to repack.c/h
223+ messages / 2 participants
[nested] [flat]
* [PATCH v45 5/8] rename cluster.c/h to repack.c/h
@ 2026-03-27 15:40 Álvaro Herrera <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Álvaro Herrera @ 2026-03-27 15:40 UTC (permalink / raw)
---
src/backend/commands/Makefile | 2 +-
src/backend/commands/matview.c | 2 +-
src/backend/commands/meson.build | 2 +-
src/backend/commands/{cluster.c => repack.c} | 6 +++---
src/backend/commands/repack_worker.c | 2 +-
src/backend/commands/tablecmds.c | 2 +-
src/backend/commands/vacuum.c | 2 +-
src/backend/libpq/pqmq.c | 2 +-
src/backend/postmaster/bgworker.c | 2 +-
src/backend/replication/logical/decode.c | 2 +-
src/backend/storage/ipc/procsignal.c | 2 +-
src/backend/tcop/postgres.c | 2 +-
src/backend/tcop/utility.c | 2 +-
src/include/commands/{cluster.h => repack.h} | 12 ++++++------
14 files changed, 21 insertions(+), 21 deletions(-)
rename src/backend/commands/{cluster.c => repack.c} (99%)
rename src/include/commands/{cluster.h => repack.h} (92%)
diff --git a/src/backend/commands/Makefile b/src/backend/commands/Makefile
index 6926bc44818..5b9d084977e 100644
--- a/src/backend/commands/Makefile
+++ b/src/backend/commands/Makefile
@@ -18,7 +18,6 @@ OBJS = \
amcmds.o \
analyze.o \
async.o \
- cluster.o \
collationcmds.o \
comment.o \
constraint.o \
@@ -51,6 +50,7 @@ OBJS = \
proclang.o \
propgraphcmds.o \
publicationcmds.o \
+ repack.o \
repack_worker.o \
schemacmds.o \
seclabel.o \
diff --git a/src/backend/commands/matview.c b/src/backend/commands/matview.c
index 539969d6eef..146a65dae8d 100644
--- a/src/backend/commands/matview.c
+++ b/src/backend/commands/matview.c
@@ -24,8 +24,8 @@
#include "catalog/namespace.h"
#include "catalog/pg_am.h"
#include "catalog/pg_opclass.h"
-#include "commands/cluster.h"
#include "commands/matview.h"
+#include "commands/repack.h"
#include "commands/tablecmds.h"
#include "commands/tablespace.h"
#include "executor/executor.h"
diff --git a/src/backend/commands/meson.build b/src/backend/commands/meson.build
index 4a0e6f0002b..9f258d566eb 100644
--- a/src/backend/commands/meson.build
+++ b/src/backend/commands/meson.build
@@ -6,7 +6,6 @@ backend_sources += files(
'amcmds.c',
'analyze.c',
'async.c',
- 'cluster.c',
'collationcmds.c',
'comment.c',
'constraint.c',
@@ -39,6 +38,7 @@ backend_sources += files(
'proclang.c',
'propgraphcmds.c',
'publicationcmds.c',
+ 'repack.c',
'repack_worker.c',
'schemacmds.c',
'seclabel.c',
diff --git a/src/backend/commands/cluster.c b/src/backend/commands/repack.c
similarity index 99%
rename from src/backend/commands/cluster.c
rename to src/backend/commands/repack.c
index 65b79ee18ff..d25ab43da7b 100644
--- a/src/backend/commands/cluster.c
+++ b/src/backend/commands/repack.c
@@ -1,6 +1,6 @@
/*-------------------------------------------------------------------------
*
- * cluster.c
+ * repack.c
* REPACK a table; formerly known as CLUSTER. VACUUM FULL also uses
* parts of this code.
*
@@ -26,7 +26,7 @@
*
*
* IDENTIFICATION
- * src/backend/commands/cluster.c
+ * src/backend/commands/repack.c
*
*-------------------------------------------------------------------------
*/
@@ -52,9 +52,9 @@
#include "catalog/pg_control.h"
#include "catalog/pg_inherits.h"
#include "catalog/toasting.h"
-#include "commands/cluster.h"
#include "commands/defrem.h"
#include "commands/progress.h"
+#include "commands/repack.h"
#include "commands/repack_internal.h"
#include "commands/tablecmds.h"
#include "commands/vacuum.h"
diff --git a/src/backend/commands/repack_worker.c b/src/backend/commands/repack_worker.c
index 35ab1a25c42..106b2b60441 100644
--- a/src/backend/commands/repack_worker.c
+++ b/src/backend/commands/repack_worker.c
@@ -20,7 +20,7 @@
#include "access/xlog_internal.h"
#include "access/xlogutils.h"
#include "access/xlogwait.h"
-#include "commands/cluster.h"
+#include "commands/repack.h"
#include "commands/repack_internal.h"
#include "libpq/pqmq.h"
#include "replication/snapbuild.h"
diff --git a/src/backend/commands/tablecmds.c b/src/backend/commands/tablecmds.c
index cc70439627b..e411d84c776 100644
--- a/src/backend/commands/tablecmds.c
+++ b/src/backend/commands/tablecmds.c
@@ -57,10 +57,10 @@
#include "catalog/storage.h"
#include "catalog/storage_xlog.h"
#include "catalog/toasting.h"
-#include "commands/cluster.h"
#include "commands/comment.h"
#include "commands/defrem.h"
#include "commands/event_trigger.h"
+#include "commands/repack.h"
#include "commands/sequence.h"
#include "commands/tablecmds.h"
#include "commands/tablespace.h"
diff --git a/src/backend/commands/vacuum.c b/src/backend/commands/vacuum.c
index 201835f30a4..ac41c819f65 100644
--- a/src/backend/commands/vacuum.c
+++ b/src/backend/commands/vacuum.c
@@ -38,9 +38,9 @@
#include "catalog/pg_database.h"
#include "catalog/pg_inherits.h"
#include "commands/async.h"
-#include "commands/cluster.h"
#include "commands/defrem.h"
#include "commands/progress.h"
+#include "commands/repack.h"
#include "commands/vacuum.h"
#include "miscadmin.h"
#include "nodes/makefuncs.h"
diff --git a/src/backend/libpq/pqmq.c b/src/backend/libpq/pqmq.c
index 1000b7bb06e..cbf6c5a004f 100644
--- a/src/backend/libpq/pqmq.c
+++ b/src/backend/libpq/pqmq.c
@@ -14,7 +14,7 @@
#include "postgres.h"
#include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack.h"
#include "libpq/libpq.h"
#include "libpq/pqformat.h"
#include "libpq/pqmq.h"
diff --git a/src/backend/postmaster/bgworker.c b/src/backend/postmaster/bgworker.c
index 1de66cc038f..40571d38d93 100644
--- a/src/backend/postmaster/bgworker.c
+++ b/src/backend/postmaster/bgworker.c
@@ -13,7 +13,7 @@
#include "postgres.h"
#include "access/parallel.h"
-#include "commands/cluster.h"
+#include "commands/repack.h"
#include "libpq/pqsignal.h"
#include "miscadmin.h"
#include "pgstat.h"
diff --git a/src/backend/replication/logical/decode.c b/src/backend/replication/logical/decode.c
index f172c0b2b40..8a969ddca41 100644
--- a/src/backend/replication/logical/decode.c
+++ b/src/backend/replication/logical/decode.c
@@ -33,7 +33,7 @@
#include "access/xlogreader.h"
#include "access/xlogrecord.h"
#include "catalog/pg_control.h"
-#include "commands/cluster.h"
+#include "commands/repack.h"
#include "replication/decode.h"
#include "replication/logical.h"
#include "replication/message.h"
diff --git a/src/backend/storage/ipc/procsignal.c b/src/backend/storage/ipc/procsignal.c
index dd980145ced..f02b9eb8312 100644
--- a/src/backend/storage/ipc/procsignal.c
+++ b/src/backend/storage/ipc/procsignal.c
@@ -19,7 +19,7 @@
#include "access/parallel.h"
#include "commands/async.h"
-#include "commands/cluster.h"
+#include "commands/repack.h"
#include "miscadmin.h"
#include "pgstat.h"
#include "port/pg_bitutils.h"
diff --git a/src/backend/tcop/postgres.c b/src/backend/tcop/postgres.c
index 4d253eddfa0..07bf300e044 100644
--- a/src/backend/tcop/postgres.c
+++ b/src/backend/tcop/postgres.c
@@ -36,10 +36,10 @@
#include "access/xact.h"
#include "catalog/pg_type.h"
#include "commands/async.h"
-#include "commands/cluster.h"
#include "commands/event_trigger.h"
#include "commands/explain_state.h"
#include "commands/prepare.h"
+#include "commands/repack.h"
#include "common/pg_prng.h"
#include "jit/jit.h"
#include "libpq/libpq.h"
diff --git a/src/backend/tcop/utility.c b/src/backend/tcop/utility.c
index 2b609bfc824..5f8c766c4be 100644
--- a/src/backend/tcop/utility.c
+++ b/src/backend/tcop/utility.c
@@ -26,7 +26,6 @@
#include "catalog/toasting.h"
#include "commands/alter.h"
#include "commands/async.h"
-#include "commands/cluster.h"
#include "commands/collationcmds.h"
#include "commands/comment.h"
#include "commands/conversioncmds.h"
@@ -46,6 +45,7 @@
#include "commands/proclang.h"
#include "commands/propgraphcmds.h"
#include "commands/publicationcmds.h"
+#include "commands/repack.h"
#include "commands/schemacmds.h"
#include "commands/seclabel.h"
#include "commands/sequence.h"
diff --git a/src/include/commands/cluster.h b/src/include/commands/repack.h
similarity index 92%
rename from src/include/commands/cluster.h
rename to src/include/commands/repack.h
index 8ed7816e7b4..6b4ebd7fa5d 100644
--- a/src/include/commands/cluster.h
+++ b/src/include/commands/repack.h
@@ -1,17 +1,17 @@
/*-------------------------------------------------------------------------
*
- * cluster.h
- * header file for postgres cluster command stuff
+ * repack.h
+ * header file for the REPACK command
*
* Portions Copyright (c) 1996-2026, PostgreSQL Global Development Group
* Portions Copyright (c) 1994-5, Regents of the University of California
*
- * src/include/commands/cluster.h
+ * src/include/commands/repack.h
*
*-------------------------------------------------------------------------
*/
-#ifndef CLUSTER_H
-#define CLUSTER_H
+#ifndef REPACK_H
+#define REPACK_H
#include <signal.h>
@@ -66,4 +66,4 @@ extern void RepackWorkerMain(Datum main_arg);
extern bool IsRepackWorker(void);
-#endif /* CLUSTER_H */
+#endif /* REPACK_H */
--
2.47.3
--cldir6umqisr673d
Content-Type: text/x-diff; charset=utf-8
Content-Disposition: attachment;
filename="v45-0006-Fix-a-few-problems-in-index-build-progress-repor.patch"
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
* [PATCH] Lock upgrade without deadlocks.
@ 2026-04-17 13:13 Antonin Houska <[email protected]>
0 siblings, 0 replies; 223+ messages in thread
From: Antonin Houska @ 2026-04-17 13:13 UTC (permalink / raw)
For REPACK (CONCURRENTLY), it's essential to upgrade ShareUpdateExclusive lock
to AccessExclusiveLock for the final stage of table processing. However, we
cannot release the earlier before requesting the latter, because if another
process ran DDL on the table in between, REPACK would have to cancel its
transaction, and waste possibly a lot of work. Without releasing
ShareUpdateExclusive lock, the upgrade can cause deadlock, and it's not really
deterministic whether REPACK or the other process will get canceled.
This patch adds a new field to the LOCK structure which tells which process is
trying to upgrade the lock - this field is set right before the upgrade. All
processes waiting in the lock queue are then woken up to check the flag.
When another process tries to get the lock after that, it checks this field,
and if it's set, it checks for deadlocks even if deadlock timeout hasn't
expired yet. If the process is already in the lock's queue and sleeping, the
lock upgrading process wakes it up so it checks the flag immediately.
At the time the upgrading process starts to sleep, all the other process
should be aware that they need to check for deadlock during their lock
acquisitions, so the upgrading process does not have to check for deadlocks
when it gets woken up. Thus it should never receive deadlock error report.
---
src/backend/commands/repack.c | 4 +-
src/backend/storage/lmgr/lmgr.c | 35 ++++++++---
src/backend/storage/lmgr/lock.c | 107 +++++++++++++++++++++++++++++++-
src/backend/storage/lmgr/proc.c | 32 +++++++---
src/include/storage/lmgr.h | 1 +
src/include/storage/lock.h | 5 +-
src/include/storage/proc.h | 4 +-
7 files changed, 166 insertions(+), 22 deletions(-)
diff --git a/src/backend/commands/repack.c b/src/backend/commands/repack.c
index 58e3867246f..310b2a65099 100644
--- a/src/backend/commands/repack.c
+++ b/src/backend/commands/repack.c
@@ -3055,8 +3055,10 @@ rebuild_relation_finish_concurrent(Relation NewHeap, Relation OldHeap,
/*
* Acquire AccessExclusiveLock on the table, its TOAST relation (if there
* is one), all its indexes, so that we can swap the files.
+ *
+ * TODO The same for indexes and TOAST?
*/
- LockRelationOid(old_table_oid, AccessExclusiveLock);
+ LockRelationOidUpgrade(old_table_oid, AccessExclusiveLock);
/*
* Lock all indexes now, not only the clustering one: all indexes need to
diff --git a/src/backend/storage/lmgr/lmgr.c b/src/backend/storage/lmgr/lmgr.c
index 2ccf7237fee..0b103485d54 100644
--- a/src/backend/storage/lmgr/lmgr.c
+++ b/src/backend/storage/lmgr/lmgr.c
@@ -58,6 +58,8 @@ typedef struct XactLockTableWaitInfo
const ItemPointerData *ctid;
} XactLockTableWaitInfo;
+static void LockRelationOidCommon(Oid relid, LOCKMODE lockmode,
+ bool isUpgrade);
static void XactLockTableWaitErrorCb(void *arg);
/*
@@ -105,6 +107,21 @@ SetLocktagRelationOid(LOCKTAG *tag, Oid relid)
*/
void
LockRelationOid(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, false);
+}
+
+/*
+ * Like above, but upgrade an existing lock.
+ */
+void
+LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode)
+{
+ LockRelationOidCommon(relid, lockmode, true);
+}
+
+static void
+LockRelationOidCommon(Oid relid, LOCKMODE lockmode, bool isUpgrade)
{
LOCKTAG tag;
LOCALLOCK *locallock;
@@ -113,7 +130,7 @@ LockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, isUpgrade);
/*
* Now that we have the lock, check for invalidation messages, so that we
@@ -157,7 +174,7 @@ ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode)
SetLocktagRelationOid(&tag, relid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -191,7 +208,7 @@ LockRelationId(LockRelId *relid, LOCKMODE lockmode)
SET_LOCKTAG_RELATION(tag, relid->dbId, relid->relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -254,7 +271,7 @@ LockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, false, true, &locallock,
- false);
+ false, false);
/*
* Now that we have the lock, check for invalidation messages; see notes
@@ -286,7 +303,7 @@ ConditionalLockRelation(Relation relation, LOCKMODE lockmode)
relation->rd_lockInfo.lockRelId.relId);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -591,7 +608,7 @@ ConditionalLockTuple(Relation relation, const ItemPointerData *tid, LOCKMODE loc
ItemPointerGetOffsetNumber(tid));
return (LockAcquireExtended(&tag, lockmode, false, true, true, NULL,
- logLockFailure) != LOCKACQUIRE_NOT_AVAIL);
+ logLockFailure, false) != LOCKACQUIRE_NOT_AVAIL);
}
/*
@@ -749,7 +766,7 @@ ConditionalXactLockTableWait(TransactionId xid, bool logLockFailure)
SET_LOCKTAG_TRANSACTION(tag, xid);
if (LockAcquireExtended(&tag, ShareLock, false, true, true, NULL,
- logLockFailure)
+ logLockFailure, false)
== LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1043,7 +1060,7 @@ ConditionalLockDatabaseObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
@@ -1123,7 +1140,7 @@ ConditionalLockSharedObject(Oid classid, Oid objid, uint16 objsubid,
objsubid);
res = LockAcquireExtended(&tag, lockmode, false, true, true, &locallock,
- false);
+ false, false);
if (res == LOCKACQUIRE_NOT_AVAIL)
return false;
diff --git a/src/backend/storage/lmgr/lock.c b/src/backend/storage/lmgr/lock.c
index c221fe96889..bcbbf36c091 100644
--- a/src/backend/storage/lmgr/lock.c
+++ b/src/backend/storage/lmgr/lock.c
@@ -810,7 +810,7 @@ LockAcquire(const LOCKTAG *locktag,
bool dontWait)
{
return LockAcquireExtended(locktag, lockmode, sessionLock, dontWait,
- true, NULL, false);
+ true, NULL, false, false);
}
/*
@@ -829,6 +829,9 @@ LockAcquire(const LOCKTAG *locktag,
*
* logLockFailure indicates whether to log details when a lock acquisition
* fails with dontWait = true.
+ *
+ * isUpgrade should be true if the backend already holds this lock in a lower
+ * mode.
*/
LockAcquireResult
LockAcquireExtended(const LOCKTAG *locktag,
@@ -837,7 +840,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure)
+ bool logLockFailure,
+ bool isUpgrade)
{
LOCKMETHODID lockmethodid = locktag->locktag_lockmethodid;
LockMethod lockMethodTable;
@@ -1119,7 +1123,8 @@ LockAcquireExtended(const LOCKTAG *locktag,
* case, because JoinWaitQueue() may discover that we can acquire the
* lock immediately after all.
*/
- waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait);
+ waitResult = JoinWaitQueue(locallock, lockMethodTable, dontWait,
+ isUpgrade);
}
if (waitResult == PROC_WAIT_STATUS_ERROR)
@@ -1225,8 +1230,85 @@ LockAcquireExtended(const LOCKTAG *locktag,
Assert(!dontWait);
PROCLOCK_PRINT("LockAcquire: sleeping on lock", proclock);
LOCK_PRINT("LockAcquire: sleeping on lock", lock, lockmode);
+
+ /*
+ * Lock upgrade can introduce deadlock, therefore enforce special
+ * behavior of other processes that deal with this lock. In
+ * particular, any waiter that sees upgradedBy set is expected to
+ * perform deadlock check as soon as it's woken up. (Since we are
+ * already in the queue, the deadlock detector has all the information
+ * it needs.)
+ */
+ if (isUpgrade)
+ {
+ dlist_iter iter;
+
+ /*
+ * There should not be multiple upgrades at the same time. XXX
+ * ERROR ?
+ */
+ Assert(lock->upgradedBy == NULL);
+
+ lock->upgradedBy = MyProc;
+
+ /*
+ * Now, before I sleep myself, wake up all the existing waiters
+ * (except for me), so they check for deadlock.
+ */
+ dclist_foreach(iter, &lock->waitProcs)
+ {
+ PGPROC *proc = dlist_container(PGPROC, waitLink,
+ iter.cur);
+
+ if (proc != MyProc)
+ SetLatch(&proc->procLatch);
+ }
+ }
LWLockRelease(partitionLock);
+ if (!isUpgrade)
+ {
+ HASH_SEQ_STATUS status;
+ LOCALLOCK *llock;
+ bool check_deadlock = false;
+
+ /*
+ * Likewise, new waiters should check the isUpgrade flag and
+ * engage deadlock detector if needed. The lock being acquired
+ * should already be in the hash.
+ *
+ * Note that, to make deadlock detection happen as soon as
+ * possible (i.e. regardless deadlock timeout), we check the other
+ * locks too. XXX There might be ways to get into the deadlock
+ * indirectly, but that needs more analysis. As long as the
+ * upgrading process skips deadlock detection altogether, the
+ * worst consequence of missing some lock wait cycle here is that
+ * the other process will be kicked-off no sooner than after its
+ * deadlock timeout has elapsed. (Which in turn means that the
+ * lock upgrade will take longer than expected.)
+ */
+ hash_seq_init(&status, LockMethodLocalHash);
+ while ((llock = (LOCALLOCK *) hash_seq_search(&status)) != NULL)
+ {
+ LOCK *mylock = llock->lock;
+
+ if (mylock && mylock->upgradedBy)
+ {
+ check_deadlock = true;
+ break;
+ }
+ }
+ /* Check for deadlock if needed. */
+ if (check_deadlock)
+ {
+ DeadLockState deadlock_state;
+
+ deadlock_state = CheckDeadLock();
+ if (deadlock_state == DS_HARD_DEADLOCK)
+ DeadLockReport();
+ }
+ }
+
waitResult = WaitOnLock(locallock, owner);
/*
@@ -1245,6 +1327,18 @@ LockAcquireExtended(const LOCKTAG *locktag,
DeadLockReport();
/* DeadLockReport() will not return */
}
+
+ /*
+ * If finishing the lock upgrade, we should not get into a deadlock
+ * anymore, so let others know that they do not have to care either.
+ */
+ if (isUpgrade)
+ {
+ LWLockAcquire(partitionLock, LW_EXCLUSIVE);
+ Assert(lock->upgradedBy != NULL);
+ lock->upgradedBy = NULL;
+ LWLockRelease(partitionLock);
+ }
}
else
LWLockRelease(partitionLock);
@@ -1322,6 +1416,7 @@ SetupLockInTable(LockMethod lockMethodTable, PGPROC *proc,
lock->nGranted = 0;
MemSet(lock->requested, 0, sizeof(int) * MAX_LOCKMODES);
MemSet(lock->granted, 0, sizeof(int) * MAX_LOCKMODES);
+ lock->upgradedBy = NULL;
LOCK_PRINT("LockAcquire: new", lock, lockmode);
}
else
@@ -1768,6 +1863,12 @@ CleanUpLock(LOCK *lock, PROCLOCK *proclock,
elog(PANIC, "proclock table corrupted");
}
+ /*
+ * Was this backend upgrading the lock?
+ */
+ if (lock->upgradedBy == MyProc)
+ lock->upgradedBy = NULL;
+
if (lock->nRequested == 0)
{
/*
diff --git a/src/backend/storage/lmgr/proc.c b/src/backend/storage/lmgr/proc.c
index 1ac25068d62..7b31370db70 100644
--- a/src/backend/storage/lmgr/proc.c
+++ b/src/backend/storage/lmgr/proc.c
@@ -95,7 +95,6 @@ static volatile sig_atomic_t got_deadlock_timeout;
static void RemoveProcFromArray(int code, Datum arg);
static void ProcKill(int code, Datum arg);
static void AuxiliaryProcKill(int code, Datum arg);
-static DeadLockState CheckDeadLock(void);
/*
@@ -1143,7 +1142,8 @@ AuxiliaryPidGetProc(int pid)
* NOTES: The process queue is now a priority queue for locking.
*/
ProcWaitStatus
-JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
+JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade)
{
LOCKMODE lockmode = locallock->tag.mode;
LOCK *lock = locallock->lock;
@@ -1230,8 +1230,12 @@ JoinWaitQueue(LOCALLOCK *locallock, LockMethod lockMethodTable, bool dontWait)
/* Must he wait for me? */
if (lockMethodTable->conflictTab[proc->waitLockMode] & myHeldLocks)
{
- /* Must I wait for him ? */
- if (lockMethodTable->conflictTab[lockmode] & proc->heldLocks)
+ /*
+ * Must I wait for him? I don't want a deadlock during lock
+ * upgrade - other processes should fail on it.
+ */
+ if ((lockMethodTable->conflictTab[lockmode] & proc->heldLocks) &&
+ !isUpgrade)
{
/*
* Yes, so we have a deadlock. Easiest way to clean up
@@ -1461,8 +1465,22 @@ ProcSleep(LOCALLOCK *locallock)
(void) WaitLatch(MyLatch, WL_LATCH_SET | WL_EXIT_ON_PM_DEATH, 0,
PG_WAIT_LOCK | locallock->tag.lock.locktag_type);
ResetLatch(MyLatch);
- /* check for deadlocks first, as that's probably log-worthy */
- if (got_deadlock_timeout)
+ /*
+ * Check for deadlocks first, as that's probably log-worthy.
+ *
+ * Do not wait for the timeout if the lock is being upgraded since
+ * the risk of deadlock is higher now. However, do not check for
+ * deadlock if the lock is being upgraded by this process - other
+ * processes should take care.
+ *
+ * TODO Possible optimization: if this is the only lock of the
+ * backend and if it did not have any weaker lock on the table so
+ * far, it should be safe to skip the deadlock check. However, to
+ * evaluate the situation, we need to take fast-path locks into
+ * account.
+ */
+ if ((got_deadlock_timeout || lock->upgradedBy) &&
+ lock->upgradedBy != MyProc)
{
deadlock_state = CheckDeadLock();
got_deadlock_timeout = false;
@@ -1819,7 +1837,7 @@ ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock)
* not, just return. If we have a real deadlock, remove ourselves from the
* lock's wait queue.
*/
-static DeadLockState
+DeadLockState
CheckDeadLock(void)
{
int i;
diff --git a/src/include/storage/lmgr.h b/src/include/storage/lmgr.h
index 2a985ce5e15..eddbe1e8622 100644
--- a/src/include/storage/lmgr.h
+++ b/src/include/storage/lmgr.h
@@ -38,6 +38,7 @@ extern void RelationInitLockInfo(Relation relation);
/* Lock a relation */
extern void LockRelationOid(Oid relid, LOCKMODE lockmode);
+extern void LockRelationOidUpgrade(Oid relid, LOCKMODE lockmode);
extern void LockRelationId(LockRelId *relid, LOCKMODE lockmode);
extern bool ConditionalLockRelationOid(Oid relid, LOCKMODE lockmode);
extern void UnlockRelationId(LockRelId *relid, LOCKMODE lockmode);
diff --git a/src/include/storage/lock.h b/src/include/storage/lock.h
index ee3cb1dc203..080b69eddfd 100644
--- a/src/include/storage/lock.h
+++ b/src/include/storage/lock.h
@@ -131,6 +131,7 @@ typedef const LockMethodData *LockMethod;
* nRequested -- total requested locks of all types.
* granted -- count of each lock type currently granted on the lock.
* nGranted -- total granted locks of all types.
+ * upgrader -- process that intends to upgrade the lock
*
* Note: these counts count 1 for each backend. Internally to a backend,
* there may be multiple grabs on a particular lock, but this is not reflected
@@ -150,6 +151,7 @@ typedef struct LOCK
int nRequested; /* total of requested[] array */
int granted[MAX_LOCKMODES]; /* counts of granted locks */
int nGranted; /* total of granted[] array */
+ PGPROC *upgradedBy; /* is lock being upgraded by this process? */
} LOCK;
#define LOCK_LOCKMETHOD(lock) ((LOCKMETHODID) (lock).tag.locktag_lockmethodid)
@@ -390,7 +392,8 @@ extern LockAcquireResult LockAcquireExtended(const LOCKTAG *locktag,
bool dontWait,
bool reportMemoryError,
LOCALLOCK **locallockp,
- bool logLockFailure);
+ bool logLockFailure,
+ bool isUpgrade);
extern void AbortStrongLockAcquire(void);
extern void MarkLockClear(LOCALLOCK *locallock);
extern bool LockRelease(const LOCKTAG *locktag,
diff --git a/src/include/storage/proc.h b/src/include/storage/proc.h
index 3e1d1fad5f9..c4bc3a24044 100644
--- a/src/include/storage/proc.h
+++ b/src/include/storage/proc.h
@@ -563,10 +563,12 @@ extern bool HaveNFreeProcs(int n, int *nfree);
extern void ProcReleaseLocks(bool isCommit);
extern ProcWaitStatus JoinWaitQueue(LOCALLOCK *locallock,
- LockMethod lockMethodTable, bool dontWait);
+ LockMethod lockMethodTable, bool dontWait,
+ bool isUpgrade);
extern ProcWaitStatus ProcSleep(LOCALLOCK *locallock);
extern void ProcWakeup(PGPROC *proc, ProcWaitStatus waitStatus);
extern void ProcLockWakeup(LockMethod lockMethodTable, LOCK *lock);
+extern DeadLockState CheckDeadLock(void);
extern void CheckDeadLockAlert(void);
extern void LockErrorCleanup(void);
extern void GetLockHoldersAndWaiters(LOCALLOCK *locallock,
--
2.47.3
--=-=-=--
^ permalink raw reply [nested|flat] 223+ messages in thread
end of thread, other threads:[~2026-04-17 13:13 UTC | newest]
Thread overview: 223+ messages (download: mbox mbox.gz follow: Atom feed)
-- links below jump to the message on this page --
2026-03-27 15:40 [PATCH v45 5/8] rename cluster.c/h to repack.c/h Álvaro Herrera <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
2026-04-17 13:13 [PATCH] Lock upgrade without deadlocks. Antonin Houska <[email protected]>
This inbox is served by agora; see mirroring instructions
for how to clone and mirror all data and code used for this inbox